# -*- coding: utf-8 -*-
import pandas as pd
import time
import os
import re
import requests
from urllib.parse import urljoin, urlparse
from bs4 import BeautifulSoup

# ======================== Global File Configuration (Isolated Data Source) ========================
KEYWORD_FILE = "非遗体育项目关键词清单.xlsx"
UNIVERSITY_LIST_FILE = "待检索高校定向名单.xlsx"
OUTPUT_RESULT = "官网定向检索_非遗体育课程数据表.xlsx"

# ======================== Network Request Configuration ========================
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Accept-Encoding": "gzip, deflate, br",
    "Connection": "keep-alive",
}
MAX_RETRY_TIMES = 2
PAGE_TIMEOUT = 12
SLEEP_BETWEEN_SCHOOL = 3.5
SLEEP_BETWEEN_PAGE = 2.2
MAX_PARSE_LINK_NUM = 5

# ======================== Pre‑defined Navigation Keyword Pool ========================
DEPARTMENT_PATH_POOL = [
    "/tyxy", "/tyb", "/sports", "/pe", "/tiyu",
    "/tyxy/index.html", "/tyb/index.htm", "/sports/home"
]
COURSE_CATEGORY_KEY = [
    "人才培养", "培养方案", "教学计划", "课程设置", "选修课",
    "公共体育", "体育课程", "课程安排", "专业课程", "培养计划",
    "特色课程", "校本课程", "本科教学", "人才培养方案"
]
INVALID_HREF_PREFIX = ("#", "javascript:", "mailto:", "tel:")


# ======================== Utility Function Module ========================
def load_keywords() -> list:
    """
    Load keyword list of intangible‑heritage sports items
    :return: cleaned keyword list
    """
    if not os.path.exists(KEYWORD_FILE):
        raise FileNotFoundError(f"Keyword file missing, check path: {KEYWORD_FILE}")
    df = pd.read_excel(KEYWORD_FILE)
    if "非遗项目名称" not in df.columns:
        raise ValueError("Required column missing: 非遗项目名称")
    keyword_list = df["非遗项目名称"].dropna().astype(str).str.strip().tolist()
    keyword_list = [item for item in keyword_list if item and item != "nan"]
    print(f"✅ Keywords loaded, total keywords: {len(keyword_list)}")
    return keyword_list


def load_target_university_list() -> list:
    """
    Load independent university retrieval list, separated from previous dataset 1‑2‑3
    :return: list of university dictionaries
    """
    if not os.path.exists(UNIVERSITY_LIST_FILE):
        raise FileNotFoundError(f"University list missing, check path: {UNIVERSITY_LIST_FILE}")
    df = pd.read_excel(UNIVERSITY_LIST_FILE)
    required_col = ["院校名称", "高校官网地址"]
    for col in required_col:
        if col not in df.columns:
            raise ValueError(f"Required column missing: {col}")
    school_dataset = []
    for _, row in df.iterrows():
        school_item = {
            "school_name": str(row["院校名称"]).strip(),
            "home_url": str(row["高校官网地址"]).strip(),
            "location": str(row.get("院校所在地", "")).strip(),
            "school_type": str(row.get("院校类型", "")).strip()
        }
        school_dataset.append(school_item)
    print(f"✅ University retrieval list loaded, total schools: {len(school_dataset)}")
    return school_dataset


def fetch_web_page(url: str) -> str | None:
    """
    Request webpage with retry mechanism
    :param url: target webpage address
    :return: html source text, None if request failed
    """
    for attempt in range(MAX_RETRY_TIMES + 1):
        try:
            response = requests.get(url, headers=HEADERS, timeout=PAGE_TIMEOUT)
            response.encoding = "utf-8"
            if response.status_code == 200:
                return response.text
            else:
                print(f"  ⚠️ Abnormal HTTP status code:{response.status_code}, retry:{attempt + 1}")
        except requests.exceptions.RequestException as err:
            print(f"  ⚠️ Network exception, retry:{attempt + 1}, error summary:{str(err)[:55]}")
        if attempt < MAX_RETRY_TIMES:
            time.sleep(2)
    return None


def build_sports_department_candidates(root_url: str) -> list:
    """
    Generate candidate PE‑department URLs based on university homepage
    :param root_url: university homepage url
    :return: candidate url list
    """
    candidate_url_list = []
    for sub_path in DEPARTMENT_PATH_POOL:
        full_link = urljoin(root_url, sub_path)
        candidate_url_list.append(full_link)
    return candidate_url_list


def scan_course_navigation_link(html_text: str, base_url: str) -> list:
    """
    Scan hyperlinks and filter curriculum‑related navigation links
    :param html_text: source text of current webpage
    :param base_url: base webpage url for relative‑path conversion
    :return: deduplicated link list [(url, anchor text),...]
    """
    soup = BeautifulSoup(html_text, "html.parser")
    raw_link_collection = []
    for a_label in soup.find_all("a", href=True):
        anchor_text = a_label.get_text(strip=True)
        href_raw = a_label["href"]
        if href_raw.startswith(INVALID_HREF_PREFIX):
            continue
        full_url = urljoin(base_url, href_raw)
        text_merge = anchor_text + " " + href_raw
        hit_flag = any(key_word in text_merge for key_word in COURSE_CATEGORY_KEY)
        if hit_flag:
            raw_link_collection.append((full_url, anchor_text))
    seen_url_set = set()
    output_link_list = []
    for link, text in raw_link_collection:
        if link not in seen_url_set:
            seen_url_set.add(link)
            output_link_list.append((link, text))
    return output_link_list


def keyword_fulltext_scan(html_text: str, keyword_array: list) -> tuple[set, list]:
    """
    Full‑text keyword scanning and extract surrounding text snippets
    :param html_text: webpage source text
    :param keyword_array: keyword library
    :return: hit‑keyword set, matched snippet list
    """
    soup = BeautifulSoup(html_text, "html.parser")
    plain_text = soup.get_text(separator=" ", strip=True)
    hit_project_set = set()
    snippet_record = []
    context_window = 35
    for keyword in keyword_array:
        index_pos = plain_text.find(keyword)
        if index_pos != -1:
            hit_project_set.add(keyword)
            start_slice = max(0, index_pos - context_window)
            end_slice = min(len(plain_text), index_pos + len(keyword) + context_window)
            fragment = plain_text[start_slice:end_slice]
            snippet_record.append({"keyword": keyword, "fragment": fragment.strip()})
    return hit_project_set, snippet_record


# ======================== Single‑university Retrieval Pipeline ========================
def single_school_retrieval(school_info: dict, keyword_list: list) -> list:
    """
    Execute complete retrieval workflow for one university: homepage‑>PE‑page‑>curriculum‑scan
    :param school_info: basic information of target university
    :param keyword_list: keyword library
    :return: retrieval records for this university
    """
    output_records = []
    s_name = school_info["school_name"]
    s_homepage = school_info["home_url"]
    s_local = school_info["location"]
    s_type = school_info["school_type"]

    home_html = fetch_web_page(s_homepage)
    if home_html is None:
        output_records.append({
            "对应非遗体育项目名称": "",
            "院校名称": s_name,
            "院校所在地": s_local,
            "院校类型": s_type,
            "非遗传统体育课程名称": "",
            "课程属性": "",
            "课程等级": "",
            "课程简介": "",
            "命中原文片段": "",
            "数据来源渠道": "高校官网定向检索",
            "备注": f"Homepage access failed, url:{s_homepage}, manual inspection required"
        })
        return output_records

    all_course_links = scan_course_navigation_link(home_html, s_homepage)
    if len(all_course_links) == 0:
        print("  Curriculum navigation not found on homepage, start detecting PE‑department pages")
        sports_sub_candidates = build_sports_department_candidates(s_homepage)
        for sub_url in sports_sub_candidates:
            sub_html = fetch_web_page(sub_url)
            if sub_html:
                print(f"  PE candidate page accessed:{sub_url}")
                sub_page_links = scan_course_navigation_link(sub_html, sub_url)
                if len(sub_page_links) > 0:
                    all_course_links.extend(sub_page_links)
                    break

    if len(all_course_links) == 0:
        output_records.append({
            "对应非遗体育项目名称": "",
            "院校名称": s_name,
            "院校所在地": s_local,
            "院校类型": s_type,
            "非遗传统体育课程名称": "",
            "课程属性": "",
            "课程等级": "",
            "课程简介": "",
            "命中原文片段": "",
            "数据来源渠道": "高校官网定向检索",
            "备注": "No curriculum‑related page detected, manual inspection required"
        })
        return output_records

    final_hit_project = set()
    all_snippet_list = []
    scan_link_limit = all_course_links[:MAX_PARSE_LINK_NUM]
    for page_url, anchor_txt in scan_link_limit:
        print(f"    Parsing curriculum page：{anchor_txt[:35]} URL:{page_url}")
        page_html = fetch_web_page(page_url)
        if page_html is None:
            continue
        hit_project, hit_snippet = keyword_fulltext_scan(page_html, keyword_list)
        final_hit_project.update(hit_project)
        all_snippet_list.extend(hit_snippet)
        time.sleep(SLEEP_BETWEEN_PAGE)

    if len(final_hit_project) > 0:
        combined_fragment_text = "; ".join([item["fragment"] for item in all_snippet_list])
        for project_name in final_hit_project:
            output_records.append({
                "对应非遗体育项目名称": project_name,
                "院校名称": s_name,
                "院校所在地": s_local,
                "院校类型": s_type,
                "非遗传统体育课程名称": project_name,
                "课程属性": "待定",
                "课程等级": "",
                "课程简介": "",
                "命中原文片段": combined_fragment_text,
                "数据来源渠道": "高校官网定向检索",
                "备注": f"Keyword matched, source page:{page_url}"
            })
    else:
        output_records.append({
            "对应非遗体育项目名称": "",
            "院校名称": s_name,
            "院校所在地": s_local,
            "院校类型": s_type,
            "非遗传统体育课程名称": "",
            "课程属性": "",
            "课程等级": "",
            "课程简介": "",
            "命中原文片段": "",
            "数据来源渠道": "高校官网定向检索",
            "备注": "Curriculum page loaded, zero keyword match, manual re‑check suggested"
        })
    return output_records


# ======================== Main Entry ========================
def main():
    """
    Main workflow of the whole retrieval task
    """
    print("=" * 75)
    print("        University Official‑Website‑Oriented Retrieval Program")
    print("=" * 75)
    try:
        keyword_dataset = load_keywords()
        university_dataset = load_target_university_list()
        global_result_list = []
        total_school_count = len(university_dataset)
        hit_school_counter = 0
        no_hit_counter = 0

        for idx, school in enumerate(university_dataset, 1):
            print(f"\n[{idx}/{total_school_count}] Retrieving university：{school['school_name']}")
            school_result = single_school_retrieval(school, keyword_dataset)
            global_result_list.extend(school_result)
            for record in school_result:
                if record["非遗传统体育课程名称"] != "":
                    hit_school_counter += 1
                    break
            else:
                no_hit_counter += 1
            time.sleep(SLEEP_BETWEEN_SCHOOL)

        df_result = pd.DataFrame(global_result_list)
        df_hit_data = df_result[df_result["非遗传统体育课程名称"] != ""].copy()
        df_empty_data = df_result[df_result["非遗传统体育课程名称"] == ""].copy()
        if not df_hit_data.empty:
            df_hit_data = df_hit_data.drop_duplicates(
                subset=["院校名称", "非遗传统体育课程名称"],
                keep="first"
            )
        df_final_output = pd.concat([df_hit_data, df_empty_data], ignore_index=True)

        df_final_output.to_excel(OUTPUT_RESULT, index=False)

        print("\n" + "=" * 75)
        print("📊 Retrieval‑task Completion Report")
        print("=" * 75)
        print(f"Total retrieved universities：{total_school_count}")
        print(f"Universities with matched courses：{hit_school_counter}")
        print(f"Zero‑hit or access‑failed universities：{no_hit_counter}")
        print(f"Total exported records after deduplication：{len(df_final_output)}")
        print(f"\n✅ Result file saved：{OUTPUT_RESULT}")
        print("\n💡 Post‑processing suggestions:")
        print("1. Filter records marked for manual inspection;")
        print("2. Distinguish news content and formally‑opened courses from matched text snippets;")
        print("3. Supplement course attribute information after browsing official webpages manually.")

    except FileNotFoundError as err:
        print(f"❌ File loading failed：{err}")
    except ValueError as err:
        print(f"❌ Dataset column validation failed：{err}")
    except Exception as err:
        print(f"❌ Unexpected runtime error：{err}")
        import traceback
        traceback.print_exc()


if __name__ == "__main__":
    main()

